# -*- coding: utf-8 -*-
"""
China National Knowledge Infrastructure (CNKI) Intangible Heritage Sports Course Literature Survey Crawler
Functions:
1. Read the excel file "清洗后_仅新增非遗体育主项目清单.xlsx" and extract cleaned‑up main‑item names as search keywords
2. Search CNKI using a fixed search‑formula template, fetch literature abstracts, and extract offering universities and course information from them
3. Export search results to an excel file in unified format

Note: Access to CNKI usually requires campus network or institutional privileges, and its anti‑crawling mechanism is strict.
      This code attempts to simulate search with requests. If data cannot be retrieved, please search manually as prompted.
"""

import pandas as pd
import time
import re
import os
import requests
from urllib.parse import quote
from bs4 import BeautifulSoup

# ========================== File Path Configuration ==========================
INPUT_FILE = "清洗后_仅新增非遗体育主项目清单.xlsx"
OUTPUT_FILE = "检索结果3_CNKI文献调研数据.xlsx"

# ========================== HTTP Request Header Configuration ==========================
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Accept-Encoding": "gzip, deflate, br",
    "Connection": "keep-alive",
    "Referer": "https://www.cnki.net/",
    "Host": "kns.cnki.net",
}

# ========================== Data Reading Module ==========================
def read_keywords():
    """Read and return the cleaned‑up main‑item keyword list"""
    if not os.path.exists(INPUT_FILE):
        raise FileNotFoundError(f"Input file not found: {INPUT_FILE}")

    df = pd.read_excel(INPUT_FILE)
    if "清洗后主项目名称" not in df.columns:
        raise ValueError(f"Required column missing in excel. Current column list: {df.columns.tolist()}")

    keywords = df["清洗后主项目名称"].dropna().astype(str).str.strip().tolist()
    keywords = [k for k in keywords if k and k != "nan"]

    if not keywords:
        raise ValueError("No valid search keywords extracted")

    print(f"✅ Successfully loaded {len(keywords)} search keywords")
    return keywords


# ========================== HTTP Request with Automatic Retry ==========================
def fetch_html(url, max_retries=2, timeout=15):
    """Send an HTTP GET request with automatic retry on failure, return the html text"""
    for attempt in range(max_retries + 1):
        try:
            response = requests.get(url, headers=HEADERS, timeout=timeout)
            response.encoding = "utf-8"
            if response.status_code == 200:
                # Detect whether a login page or captcha page is returned (judged by title content)
                if "登录" in response.text or "验证码" in response.text:
                    print("  ⚠️ Page requires login or captcha. Please confirm the network environment (campus network) or search manually.")
                    return None
                return response.text
            else:
                print(f"  ⚠️ HTTP {response.status_code} (Attempt {attempt+1}/{max_retries+1})")
        except requests.exceptions.RequestException as e:
            print(f"  ⚠️ Request Exception (Attempt {attempt+1}/{max_retries+1}): {str(e)[:50]}")
        if attempt < max_retries:
            time.sleep(3)
    return None


# ========================== CNKI Search Function ==========================
def search_cnki(query, max_pages=2):
    """
    Search the given search formula on CNKI and return a list of articles (title, abstract, source, etc.)
    query: url‑encoded search formula
    """
    # Use the default CNKI search interface (simplified version)
    # Note: actual CNKI search may require POST; GET is used here, and some parameters may need adjustment
    base_url = "https://kns.cnki.net/kns8/DefaultResult/Index"
    params = {
        "korder": "1",         # Sorting method
        "sel": "1",            # Selection scope
        "sfield": "all",       # Search field
        "q": query,            # Search keyword
        "p": "1",              # Page number (fetch multiple pages via loop)
    }

    all_articles = []
    for page in range(1, max_pages + 1):
        params["p"] = str(page)
        # Build the complete url
        url = base_url + "?" + "&".join([f"{k}={quote(str(v))}" for k, v in params.items()])
        print(f"  🌐 Requesting page {page}: {url[:100]}...")

        html = fetch_html(url)
        if not html:
            break

        # Parse html to extract literature information
        soup = BeautifulSoup(html, "html.parser")
        # CNKI literature list is usually wrapped in <table class="ResultTable"> or <div class="list-item">
        # Try multiple selectors
        items = soup.select("table.ResultTable tr") or soup.select(".list-item")
        if not items:
            # Check for empty search result
            if "没有检索到" in html or "无结果" in html:
                print("  ⚠️ Search returned no results")
                break
            # Page structure may differ, print partial text for debugging
            # print(html[:500])
            break

        page_articles = 0
        for item in items:
            # Extract title
            title_elem = item.find("a", class_="fz14") or item.find("a", class_="title")
            if not title_elem:
                continue
            title = title_elem.get_text(strip=True)

            # Extract abstract (may exist in <p class="abstract">)
            abstract_elem = item.find("p", class_="abstract") or item.find("div", class_="abstract")
            abstract = abstract_elem.get_text(strip=True) if abstract_elem else ""

            # Extract source (journal / university)
            source_elem = item.find("p", class_="source") or item.find("div", class_="source")
            source = source_elem.get_text(strip=True) if source_elem else ""

            if title:
                all_articles.append({
                    "title": title,
                    "abstract": abstract,
                    "source": source
                })
                page_articles += 1

        print(f"  ✅ Page {page}: {page_articles} articles fetched")
        if page_articles == 0:
            break
        time.sleep(3)  # Interval between pages

    return all_articles


# ========================== Extract Institutions and Courses from Literature ==========================
def extract_institution(text):
    """Extract university names from text (prioritize matching 'XX大学', 'XX学院', etc.)"""
    # Matching pattern: Chinese university / college / sports college, etc.
    pattern = r'([\u4e00-\u9fa5]{2,10}(?:大学|学院|体育学院|师范大学|民族大学|职业学院|高等专科学校))'
    matches = re.findall(pattern, text)
    # Deduplicate and filter by length
    institutions = []
    for m in matches:
        if len(m) >= 4 and m not in institutions:
            institutions.append(m)
    return institutions


def extract_courses(text, keywords):
    """Match intangible heritage sports course keywords from text"""
    matched = []
    for kw in keywords:
        # Check whether keyword appears in text (as independent word)
        # Simple substring matching
        if kw in text:
            matched.append(kw)
    return matched


def infer_university_type(name):
    """Infer university type from its name"""
    if any(k in name for k in ["体育学院", "体育大学", "体育职业"]):
        return "体育类"
    if any(k in name for k in ["师范大学", "师范学院", "师范高等专科"]):
        return "师范类"
    if any(k in name for k in ["民族大学", "民族学院"]):
        return "民族类"
    if any(k in name for k in ["大学", "学院"]):
        return "综合类"
    return "综合类"


def infer_location(name):
    """Simply infer location from the name prefix (only use first two characters as province abbreviation; imprecise but usable as reference)"""
    # Common province abbreviation mapping
    province_map = {
        "北京": "北京", "上海": "上海", "天津": "天津", "重庆": "重庆",
        "河北": "河北", "山西": "山西", "辽宁": "辽宁", "吉林": "吉林",
        "黑龙江": "黑龙江", "江苏": "江苏", "浙江": "浙江", "安徽": "安徽",
        "福建": "福建", "江西": "江西", "山东": "山东", "河南": "河南",
        "湖北": "湖北", "湖南": "湖南", "广东": "广东", "海南": "海南",
        "四川": "四川", "贵州": "贵州", "云南": "云南", "陕西": "陕西",
        "甘肃": "甘肃", "青海": "青海", "台湾": "台湾", "内蒙古": "内蒙古",
        "广西": "广西", "西藏": "西藏", "宁夏": "宁夏", "新疆": "新疆",
    }
    for key in province_map:
        if name.startswith(key):
            return province_map[key]
    return ""


# ========================== Main Program Entry ==========================
def main():
    print("=" * 70)
    print("China National Knowledge Infrastructure (CNKI) Intangible Heritage Sports Course Literature Survey Crawler")
    print("Note: This program requires campus network or institutional access privileges. Search manually if data cannot be retrieved.")
    print("=" * 70)

    try:
        # ---------- Step 1: Load search keywords ----------
        keywords = read_keywords()
        print(f"Keyword preview (first 10 items): {keywords[:10]}")

        # ---------- Step 2: Build search formulas and search them in sequence ----------
        # Fixed search‑formula templates
        templates = [
            ("(非遗传统体育) AND (高校课程开设) AND (院校)", "Global Search"),
            ("(民族民间体育) AND (高校公共体育选修课)", "Global Search"),
        ]
        # Per‑keyword search formula
        per_keyword_template = "({}) AND (高校教学)"

        all_articles = []
        search_queries = []

        # Execute the two global searches first
        for query, desc in templates:
            print(f"\n🔍 Executing {desc}: {query}")
            search_queries.append((query, desc))
            articles = search_cnki(query, max_pages=1)  # Global search only fetches 1 page
            if articles:
                all_articles.extend(articles)
                print(f"  ✅ Total {len(articles)} articles fetched")
            else:
                print(f"  ⚠️ No articles fetched, manual search may be required")
            time.sleep(3)

        # Then search for each keyword
        print("\n" + "=" * 70)
        print("🔍 Starting per‑keyword sequential search")
        print("=" * 70)

        total = len(keywords)
        for idx, kw in enumerate(keywords, 1):
            query = per_keyword_template.format(kw)
            print(f"\n[{idx}/{total}] Search keyword: {kw}")
            print(f"  Search formula: {query}")
            articles = search_cnki(query, max_pages=1)  # Each keyword only fetches 1 page
            if articles:
                all_articles.extend(articles)
                print(f"  ✅ {len(articles)} articles fetched")
            else:
                print(f"  ⚠️ No articles fetched, manual search recommended")
            time.sleep(3)

        # ---------- Step 3: Extract institutions and courses from literature ----------
        print("\n" + "=" * 70)
        print("📊 Starting extraction of institution and course information from literature abstracts")
        print("=" * 70)

        results = []
        for article in all_articles:
            text = article["title"] + " " + article["abstract"] + " " + article["source"]
            institutions = extract_institution(text)
            courses = extract_courses(text, keywords)

            # Skip if neither institution nor course is present
            if not institutions or not courses:
                continue

            # Generate one record for each institution‑course combination
            for inst in institutions:
                for course in courses:
                    results.append({
                        "对应非遗体育项目名称": course,
                        "院校名称": inst,
                        "院校所在地": infer_location(inst),
                        "院校类型": infer_university_type(inst),
                        "非遗传统体育课程名称": course,
                        "课程属性": "",   # Not provided on CNKI
                        "课程等级": "",   # Not provided on CNKI
                        "课程简介": article["abstract"][:200] + "..." if len(article["abstract"]) > 200 else article["abstract"],
                        "数据来源渠道": "中国知网CNKI文献",
                        "备注": f"文献标题：{article['title']}"
                    })

        # ---------- Step 4: Deduplicate ----------
        if results:
            df_results = pd.DataFrame(results)
            # Deduplicate by university name + intangible heritage sports course name
            df_results = df_results.drop_duplicates(
                subset=["院校名称", "非遗传统体育课程名称"],
                keep="first"
            )
        else:
            df_results = pd.DataFrame(columns=[
                "对应非遗体育项目名称", "院校名称", "院校所在地", "院校类型",
                "非遗传统体育课程名称", "课程属性", "课程等级", "课程简介",
                "数据来源渠道", "备注"
            ])

        # ---------- Step 5: Export to excel ----------
        df_results.to_excel(OUTPUT_FILE, index=False)

        # ---------- Step 6: Print statistical report ----------
        print("\n" + "=" * 70)
        print("📊 Crawling‑task Completion Statistics")
        print("=" * 70)
        print(f"Total search keywords: {len(keywords)}")
        print(f"Total articles fetched: {len(all_articles)}")
        print(f"Valid records extracted: {len(results)}")
        print(f"Deduplicated records: {len(df_results)}")
        print(f"\n✅ Output file: {OUTPUT_FILE}")
        print("=" * 70)

        if not df_results.empty:
            print("\n📖 Result preview (first 10 entries):")
            print("-" * 70)
            preview_cols = ["对应非遗体育项目名称", "院校名称", "院校类型", "非遗传统体育课程名称"]
            print(df_results[preview_cols].head(10).to_string(index=False))
            print("-" * 70)

            # Count occurrence of each course
            print("\n📊 Number of universities mentioning each intangible heritage sports project in literature:")
            course_count = df_results.groupby("对应非遗体育项目名称").size().sort_values(ascending=False)
            for course, count in course_count.head(15).items():
                print(f"  {course}: {count} universities")
            if len(course_count) > 15:
                print(f"  ... {len(course_count) - 15} more projects omitted")
        else:
            print("\n⚠️ No valid data extracted")
            print("Potential reasons:")
            print("1. CNKI search returned no related literature")
            print("2. Literature abstracts do not explicitly mention institution or course names")
            print("3. Anti‑crawling mechanism prevented successful content retrieval")
            print("Suggestion: search manually on CNKI with the following formulas and supplement the data:")
            print("  - (非遗传统体育) AND (高校课程开设) AND (院校)")
            print("  - (民族民间体育) AND (高校公共体育选修课)")
            print("  - (具体项目名称) AND (高校教学)")

    except FileNotFoundError as e:
        print(f"❌ File‑related error: {e}")
    except ValueError as e:
        print(f"❌ Data‑validation error: {e}")
    except Exception as e:
        print(f"❌ Unexpected runtime error: {e}")
        import traceback
        traceback.print_exc()


if __name__ == "__main__":
    main()
